Skip to content

RubricsAsJudge reward

Rubric

Bases: DataModel

One rubric criterion for RubricsAsJudge.

Source code in synalinks/src/rewards/rubrics_as_judge.py
class Rubric(DataModel):
    """One rubric criterion for `RubricsAsJudge`."""

    name: str = Field(
        description=(
            "Short identifier used as the JSON property name. It is normalized "
            "to snake_case."
        ),
    )
    description: str = Field(
        description="Description of what the judge should evaluate.",
    )
    weight: float = Field(
        default=1.0,
        description="Positive relative weight.",
    )

    @field_validator("name")
    @classmethod
    def _normalize_name(cls, name):
        normalized = normalize_rubric_name(name)
        if not RUBRIC_NAME_REGEX.match(normalized):
            raise ValueError(
                f"Rubric name {name!r} normalizes to {normalized!r}, "
                "which is not usable as a JSON property."
            )
        return normalized

    @field_validator("weight")
    @classmethod
    def _check_weight(cls, weight):
        if weight <= 0:
            raise ValueError(f"Rubric weight is {weight}; weights must be positive.")
        return weight

RubricsAsJudge

Bases: ProgramAsJudge

Grade an answer against named criteria and combine them by weight.

Parameters:

Name Type Description Default
language_model LanguageModel

The language model that grades.

None
rubrics list | str

Criteria. Each item can be a Rubric, a dict with name, description and optional weight, or a bare string. A string selects a built-in rubric preset (see synalinks.rewards.list_rubrics).

None
prompt_template str

Optional Jinja2 prompt template.

None
examples list

Optional examples for the prompt.

None
instructions str

Optional judge instructions.

None
score_type type | str

Per-criterion score scale. Defaults to Rating20. Language model only.

None
reduction str

Reward reduction. Defaults to "mean".

'mean'
name str

Optional reward name.

'rubrics_as_judge'
in_mask list

Optional fields to keep before judging.

None
out_mask list

Optional fields to remove before judging.

None
in_mask_pattern str

Optional regex of fields to keep.

None
out_mask_pattern str

Optional regex of fields to remove.

None
decision_model DecisionModel

Optional. A decision model to grade with instead of the language model: every criterion is a score question over the same ordered levels, all in one call. It does not write a critique, so the critique lists the level and confidence of each criterion instead.

None
Source code in synalinks/src/rewards/rubrics_as_judge.py
@synalinks_export(
    [
        "synalinks.RubricsAsJudge",
        "synalinks.rewards.RubricsAsJudge",
    ]
)
class RubricsAsJudge(ProgramAsJudge):
    """Grade an answer against named criteria and combine them by weight.

    Args:
        language_model (LanguageModel): The language model that grades.
        rubrics (list | str): Criteria. Each item can be a `Rubric`, a dict with
            `name`, `description` and optional `weight`, or a bare string. A string
            selects a built-in rubric preset (see `synalinks.rewards.list_rubrics`).
        prompt_template (str): Optional Jinja2 prompt template.
        examples (list): Optional examples for the prompt.
        instructions (str): Optional judge instructions.
        score_type (type | str): Per-criterion score scale. Defaults to `Rating20`.
            Language model only.
        reduction (str): Reward reduction. Defaults to `"mean"`.
        name (str): Optional reward name.
        in_mask (list): Optional fields to keep before judging.
        out_mask (list): Optional fields to remove before judging.
        in_mask_pattern (str): Optional regex of fields to keep.
        out_mask_pattern (str): Optional regex of fields to remove.
        decision_model (DecisionModel): Optional. A decision model to grade
            with instead of the language model: every criterion is a score
            question over the same ordered levels, all in one call. It does
            not write a critique, so the `critique` lists the level and
            confidence of each criterion instead.
    """

    def __init__(
        self,
        language_model=None,
        rubrics=None,
        prompt_template=None,
        examples=None,
        instructions=None,
        score_type=None,
        reduction="mean",
        name="rubrics_as_judge",
        in_mask=None,
        out_mask=None,
        in_mask_pattern=None,
        out_mask_pattern=None,
        decision_model=None,
    ):
        program = RubricsAsJudgeProgram(
            language_model=language_model,
            rubrics=rubrics,
            prompt_template=prompt_template,
            examples=examples,
            instructions=instructions,
            score_type=score_type,
            decision_model=decision_model,
        )
        super().__init__(
            program=program,
            reduction=reduction,
            name=name,
            in_mask=in_mask,
            out_mask=out_mask,
            in_mask_pattern=in_mask_pattern,
            out_mask_pattern=out_mask_pattern,
        )

    @classmethod
    def from_config(cls, config):
        program = serialization_lib.deserialize_synalinks_object(config.pop("program"))
        kwargs = {
            "language_model": program.language_model,
            "prompt_template": program.prompt_template,
            "examples": program.examples,
            "instructions": program.instructions,
            "score_type": program.score_type,
            **config,
        }
        if program.decision_model is not None:
            kwargs["decision_model"] = program.decision_model
        if cls is RubricsAsJudge:
            kwargs["rubrics"] = [rubric.get_json() for rubric in program.rubrics]
        return cls(**kwargs)

RubricsAsJudgeProgram

Bases: Module

Score each criterion with a language (or decision) model, then combine in Python.

Source code in synalinks/src/rewards/rubrics_as_judge.py
class RubricsAsJudgeProgram(Module):
    """Score each criterion with a language (or decision) model, then combine
    in Python."""

    def __init__(
        self,
        language_model=None,
        rubrics=None,
        prompt_template=None,
        examples=None,
        instructions=None,
        score_type=None,
        name=None,
        description=None,
        trainable=True,
        decision_model=None,
    ):
        super().__init__(name=name, description=description, trainable=trainable)
        if isinstance(rubrics, str):
            rubrics = get_rubric(rubrics)
        if not rubrics:
            raise ValueError("`rubrics` is required and must not be empty.")
        self.rubrics = [parse_rubric(rubric) for rubric in rubrics]
        names = [rubric.name for rubric in self.rubrics]
        if len(set(names)) != len(names):
            raise ValueError(f"Duplicate rubric names after normalization: {names}.")

        self.language_model = language_model
        self.decision_model = resolve_decision_model(decision_model, language_model)
        if decision_model is None and score_type is not None:
            # A score scale is for a language model: the default decision
            # model does not apply.
            self.decision_model = None
        self.prompt_template = prompt_template
        self.examples = examples
        prompt_variables = {"rubrics": [rubric.get_json() for rubric in self.rubrics]}
        if self.decision_model is not None:
            if score_type is not None:
                raise ValueError(
                    "`score_type` does not apply with a `DecisionModel`: every "
                    "criterion is graded on the same ordered levels."
                )
            self.score_type = None
            self.instructions = instructions or DEFAULT_DECISION_MODEL_INSTRUCTIONS
            self.schema = rubrics_decision_model_schema(self.rubrics)
            self.output_schema = rubrics_schema_with_reward(
                rubrics_decision_model_output_schema(self.rubrics)
            )
        else:
            self.score_type = get_score_type(score_type or Rating20)
            self.instructions = instructions or DEFAULT_INSTRUCTIONS
            self.schema = rubrics_schema(self.rubrics, self.score_type)
            self.output_schema = rubrics_schema_with_reward(self.schema)
            prompt_variables["score_scale"] = score_type_description(self.score_type)
        self.generator = Generator(
            schema=self.schema,
            language_model=language_model,
            decision_model=self.decision_model,
            prompt_template=prompt_template or default_rubrics_prompt_template(),
            prompt_variables=prompt_variables,
            examples=examples,
            instructions=self.instructions,
            name="generator_" + self.name,
        )

    async def call(self, inputs):
        if not isinstance(inputs, (list, tuple)):
            raise ValueError("The inputs should be a list or tuple.")
        if len(inputs) != 2:
            raise ValueError("The inputs of the program should have a length of 2.")
        y_true, y_pred = inputs
        if not y_pred:
            return CritiqueWithReward(
                critique="Empty prediction: nothing to evaluate.", reward=0.0
            ).to_json_data_model()

        if y_true:
            y_true = await ops.prefix(y_true, prefix="gold", name="gold_y_true")
            inputs = await ops.concat(y_true, y_pred, name="y_true_with_y_pred")
        else:
            inputs = y_pred
        grades = await self.generator(inputs)
        if grades is None:
            return CritiqueWithReward(
                critique="The judge produced no grade sheet.", reward=0.0
            ).to_json_data_model()
        return self.combine(grades)

    def combine(self, grades):
        if is_symbolic_data_model(grades):
            return SymbolicDataModel(schema=self.output_schema, name=grades.name)

        json = dict(grades.get_json())
        if self.score_type is None:
            json = self.decision_model_grades(json)
        total = 0.0
        weights = 0.0
        missing = []
        for rubric in self.rubrics:
            value = json.get(rubric.name)
            if value is None:
                missing.append(rubric.name)
                continue
            if self.score_type is not None:
                value = normalize_score(value, self.score_type)
            total += value * rubric.weight
            weights += rubric.weight
        if missing:
            return CritiqueWithReward(
                critique=(
                    "The judge skipped "
                    f"{', '.join(missing)}; the grade sheet is incomplete."
                ),
                reward=0.0,
            ).to_json_data_model()
        json["reward"] = total / weights if weights else 0.0
        return JsonDataModel(json=json, schema=self.output_schema, name=grades.name)

    def decision_model_grades(self, answers):
        """Turn a decision model's score answers into normalized grades."""
        json = {}
        critique = []
        for rubric in self.rubrics:
            answer = answers.get(rubric.name)
            if answer is None:
                continue
            top = len(answer["legend"]) - 1
            normalized = answer["score"] / top
            level = answer["legend"][str(int(answer["score"] + 0.5))]
            critique.append(
                f"{rubric.name}: {level} (score {normalized:.2f}, "
                f"confidence {answer['confidence']:.2f})"
            )
            json[rubric.name] = normalized
        return {"critique": "\n".join(critique), **json}

    def get_config(self):
        config = {
            "rubrics": [rubric.get_json() for rubric in self.rubrics],
            "prompt_template": self.prompt_template,
            "examples": self.examples,
            "instructions": self.instructions,
            "name": self.name,
            "description": self.description,
            "trainable": self.trainable,
        }
        if self.score_type is not None:
            config["score_type"] = serialize_score_type(self.score_type)
        if self.decision_model is not None:
            config["decision_model"] = serialization_lib.serialize_synalinks_object(
                self.decision_model
            )
        return {
            "language_model": serialization_lib.serialize_synalinks_object(
                self.language_model
            ),
            **config,
        }

    @classmethod
    def from_config(cls, config):
        language_model = serialization_lib.deserialize_synalinks_object(
            config.pop("language_model")
        )
        if "decision_model" in config:
            config["decision_model"] = serialization_lib.deserialize_synalinks_object(
                config.pop("decision_model")
            )
        return cls(language_model=language_model, **config)

decision_model_grades(answers)

Turn a decision model's score answers into normalized grades.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def decision_model_grades(self, answers):
    """Turn a decision model's score answers into normalized grades."""
    json = {}
    critique = []
    for rubric in self.rubrics:
        answer = answers.get(rubric.name)
        if answer is None:
            continue
        top = len(answer["legend"]) - 1
        normalized = answer["score"] / top
        level = answer["legend"][str(int(answer["score"] + 0.5))]
        critique.append(
            f"{rubric.name}: {level} (score {normalized:.2f}, "
            f"confidence {answer['confidence']:.2f})"
        )
        json[rubric.name] = normalized
    return {"critique": "\n".join(critique), **json}

default_rubrics_prompt_template()

Return the default prompt template used by RubricsAsJudge.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def default_rubrics_prompt_template():
    """Return the default prompt template used by `RubricsAsJudge`."""
    return """
<instructions>
{{ instructions }}
</instructions>
{% if rubrics %}
<rubrics>
{% for rubric in rubrics %}
<rubric>
<name>
{{ rubric.name }}
</name>
<description>
{{ rubric.description }}
</description>{% if rubric.weight != 1.0 %}
<weight>
{{ "%g"|format(rubric.weight) }}
</weight>{% endif %}
</rubric>
{% endfor %}
</rubrics>
{% endif %}{% if score_scale %}
<score_scale>
{{ score_scale }}
</score_scale>
{% endif %}{% if inputs_schema %}
<input_schema>
{{ inputs_schema }}
</input_schema>
{% endif %}{% if outputs_schema %}
<output_schema>
{{ outputs_schema }}
</output_schema>
{% endif %}{% if examples %}
<examples>
{% for example in examples %}
<example>
<input>
{{ example[0] }}
</input>
<output>
{{ example[1] }}
</output>
</example>
{% endfor %}
</examples>
{% endif %}
""".strip()

normalize_rubric_name(name)

Normalize a rubric name into a JSON property name.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def normalize_rubric_name(name):
    """Normalize a rubric name into a JSON property name."""
    name = to_snake_case(str(name).strip())
    if not name or name[0].isdigit():
        name = f"criterion_{name}" if name else "criterion"
    return name

parse_rubric(rubric)

Return a Rubric from a Rubric, a dict or a plain description.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def parse_rubric(rubric):
    """Return a `Rubric` from a `Rubric`, a dict or a plain description."""
    if isinstance(rubric, Rubric):
        return rubric
    if isinstance(rubric, dict):
        return Rubric(**rubric)
    if isinstance(rubric, str):
        return Rubric(name=normalize_rubric_name(rubric)[:40], description=rubric)
    raise TypeError(f"Cannot interpret {rubric!r} as a rubric.")

rubrics_decision_model_output_schema(rubrics)

Return the grade-sheet schema of a DecisionModel judge.

Each criterion is its score normalized to [0, 1], next to a critique that lists the level each criterion landed on.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def rubrics_decision_model_output_schema(rubrics):
    """Return the grade-sheet schema of a `DecisionModel` judge.

    Each criterion is its score normalized to [0, 1], next to a critique that
    lists the level each criterion landed on.
    """
    properties = {
        "critique": {
            "title": "Critique",
            "type": "string",
            "description": "The level and confidence of each criterion.",
        }
    }
    for rubric in rubrics:
        properties[rubric.name] = {
            "title": rubric.name,
            "type": "number",
            "minimum": 0.0,
            "maximum": 1.0,
            "description": rubric.description,
        }
    return {
        "title": "RubricGrades",
        "type": "object",
        "properties": properties,
        "required": list(properties),
    }

rubrics_decision_model_schema(rubrics)

Return the grade-sheet schema a DecisionModel answers.

Decision models do not write text, so there is no critique: each criterion is a score question over DECISION_MODEL_LEVELS.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def rubrics_decision_model_schema(rubrics):
    """Return the grade-sheet schema a `DecisionModel` answers.

    Decision models do not write text, so there is no `critique`: each
    criterion is a score question over `DECISION_MODEL_LEVELS`.
    """
    properties = {
        rubric.name: score_schema(
            f"How well does the answer meet this criterion: {rubric.description}",
            DECISION_MODEL_LEVELS,
        )
        for rubric in rubrics
    }
    return {
        "title": "RubricGrades",
        "type": "object",
        "properties": properties,
        "required": list(properties),
        "additionalProperties": False,
    }

rubrics_schema(rubrics, score_type)

Return the constrained grade-sheet schema for a list of rubrics.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def rubrics_schema(rubrics, score_type):
    """Return the constrained grade-sheet schema for a list of rubrics."""
    score_type = get_score_type(score_type)
    properties = {
        "critique": {
            "title": "Critique",
            "type": "string",
            "description": "Why each criterion was scored the way it was.",
        }
    }
    for rubric in rubrics:
        properties[rubric.name] = {
            "title": rubric.name,
            "type": score_type_json_type(score_type),
            "enum": [member.value for member in score_type],
            "description": rubric.description,
        }
    return {
        "title": "RubricGrades",
        "type": "object",
        "properties": properties,
        "required": ["critique", *[rubric.name for rubric in rubrics]],
    }

rubrics_schema_with_reward(schema)

Return a copy of the grade-sheet schema with the computed reward field.

Source code in synalinks/src/rewards/rubrics_as_judge.py
def rubrics_schema_with_reward(schema):
    """Return a copy of the grade-sheet schema with the computed reward field."""
    schema = copy.deepcopy(schema)
    schema.setdefault("properties", {})["reward"] = {
        "title": "Reward",
        "type": "number",
        "minimum": 0.0,
        "maximum": 1.0,
        "description": "The weighted mean of the normalized criterion scores.",
    }
    required = list(schema.get("required", []))
    if "reward" not in required:
        required.append("reward")
    schema["required"] = required
    return schema